#!/usr/bin/env python

import os
import re
import time
from bs4 import BeautifulSoup

start = time.time()

def search_words(root_dir, words, output_file):
    if os.path.exists(output_file):
        os.remove(output_file)
    html = "<html><body><h1>Search Results</h1>"
    html += "<meta http-equiv='content-type' content='text/html; charset=UTF-8'>"
    html += "<meta charset='utf-8'>"
    html += "<script>function copySentence(event) {navigator.clipboard.writeText(event.target.textContent);}</script>"
    html += "<div id='google_translate_element'></div>"
    html += "<script type='text/javascript'>function googleTranslateElementInit() {new google.translate.TranslateElement({pageLanguage: 'hi', layout: google.translate.TranslateElement.InlineLayout.SIMPLE}, 'google_translate_element');}</script>"
    html += "<script type='text/javascript' src='//translate.google.com/translate_a/element.js?cb=googleTranslateElementInit'></script>"
    html += "<ul>"
    words = [word.strip() for word in words.split(',')]
    results = []
    output_dir = os.path.dirname(__file__)
    search_dir = os.path.join(output_dir)
    if not os.path.exists(search_dir):
        os.makedirs(search_dir)
    for root, dirs, files in os.walk(root_dir):
        # if'search' in root:  # replace 'dir_to_exclude' with the actual directory name
            # continue
        for file in files:
            if file.endswith('.html'):
                file_path = os.path.join(root, file)
                relative_path = os.path.relpath(file_path, output_dir)
                relative_path = './' + relative_path
                try:
                    with open(file_path, 'r') as f:
                        content = f.read()
                        soup = BeautifulSoup(content, 'html.parser')
                        title = soup.find('title')
                        if title:
                            title_text = title.text.strip()
                            title_text = title_text.replace('(गायत्री परिवार - Gayatri Pariwar)', '')
                        else:
                            title_text = file
                        p_elements = soup.find_all('p')
                        for p in p_elements:
                            p_text = p.get_text()
                            p_words = p_text.split()
                            indices = []
                            for word in words:
                                indices.append([i for i, x in enumerate(p_words) if x.lower().find(word.lower())!= -1])
                            for idx_list in indices[0]:
                                if all(any(abs(idx_list - idx) <= 100 for idx in idxs) for idxs in indices[1:]):
                                    start = idx_list - 9
                                    end = idx_list + 9 + 1
                                    if start < 0:
                                        start = 0
                                    if end > len(p_words):
                                        end = len(p_words)
                                    snippet =' '.join(p_words[start:end])
                                    bolded_snippet = snippet
                                    for word in words:
                                        bolded_snippet = bolded_snippet.replace(word, f"<b>{word}</b>")
                                    results.append((title_text, relative_path, bolded_snippet, len([w for w in words if w in bolded_snippet]), p_text))
                except Exception as e:
                    print(f"Error reading {file_path}: {e}")
    exact_matches = [result for result in results if result[3] == len(words)]
    exact_matches.sort(key=lambda x: tuple(x[2].lower().find(word.lower()) for word in words))
    exact_matches = exact_matches[:100]
    for title_text, relative_path, snippet, _, p_text in exact_matches:
        title = snippet +' '+ p_text[p_text.find(snippet) + len(snippet):].strip()
        stop_chars = ['।','?','॥','.']
        title_end_index = len(snippet) + 315
        for char in stop_chars:
            index = title[:title_end_index].rfind(char)
            if index!= -1:
                title_end_index = index + 1
                break
        title = title[:title_end_index].strip()
        title = re.sub('<.*?>', '', title)
        html += f"<li><a href='{relative_path}' target='_blank' title='{title}' onclick='copySentence(event)'>{title_text}</a> - {snippet}</li>"
    html += "</ul><h2>Partial Matches</h2><ul>"
    partial_matches = [result for result in results if result[3]!= len(words)]
    partial_matches.sort(key=lambda x: sum([words.index(word) for word in x[2].split() if word in words]), reverse=True)
    partial_matches = partial_matches[:5]
    for title_text, relative_path, snippet, _, p_text in partial_matches:
        title = snippet +' '+ p_text[p_text.find(snippet) + len(snippet):].strip()
        stop_chars = ['।','?','॥','.']
        title_end_index = len(snippet) + 315
        for char in stop_chars:
            index = title[:title_end_index].rfind(char)
            if index!= -1:
                title_end_index = index + 1
                break
        title = title[:title_end_index].strip()
        html += f"<li><a href='{relative_path}' target='_blank' title='{title}' onclick='copySentence(event)'>{title_text}</a> - {snippet}</li>"
    html += "</ul></body></html>"
    output_path = os.path.join(search_dir, output_file)
    with open(output_path, 'w') as f:
        f.write(html)
    print(html)

root_dir = '../'
words = 'अध्यात्म,क्या,है'
output_file ='search_multiple_words.html'
# search_words(root_dir, words, output_file)

print("See "+output_file)

end = time.time()
print("Elapsed (with compilation) = %s" % (end - start))
